officeparser 6.1.1 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +301 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +74 -31
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +828 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +783 -5
  54. package/dist/types.js +73 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.d.ts +8 -3
  60. package/dist/utils/envUtils.js +117 -34
  61. package/dist/utils/errorUtils.d.ts +17 -29
  62. package/dist/utils/errorUtils.js +110 -52
  63. package/dist/utils/moduleLoader.js +19 -11
  64. package/dist/utils/ocrUtils.js +2 -1
  65. package/dist/utils/sheetUtils.d.ts +7 -0
  66. package/dist/utils/sheetUtils.js +35 -0
  67. package/dist/utils/styleMapper.d.ts +36 -0
  68. package/dist/utils/styleMapper.js +224 -0
  69. package/dist/utils/xmlUtils.d.ts +0 -8
  70. package/dist/utils/xmlUtils.js +2 -1
  71. package/package.json +26 -7
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
- # officeParser 📄🚀
1
+ # officeParser 📄🚀 - The Most Versatile Office Parser & Generator
2
2
 
3
- A robust, strictly-typed Node.js and Browser library for parsing office files ([`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format)). It produces a clean, hierarchical Abstract Syntax Tree (AST) with rich metadata, text formatting, and full attachment support.
3
+ A robust, strictly-typed Node.js and Browser library for parsing and generating office files. It not only extracts content from [`docx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`pptx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`xlsx`](https://en.wikipedia.org/wiki/Office_Open_XML), [`odt`](https://en.wikipedia.org/wiki/OpenDocument), [`odp`](https://en.wikipedia.org/wiki/OpenDocument), [`ods`](https://en.wikipedia.org/wiki/OpenDocument), [`pdf`](https://en.wikipedia.org/wiki/PDF), [`rtf`](https://en.wikipedia.org/wiki/Rich_Text_Format), [`csv`](https://en.wikipedia.org/wiki/Comma-separated_values), [`md`](https://en.wikipedia.org/wiki/Markdown), and [`html`](https://en.wikipedia.org/wiki/HTML) into a rich Abstract Syntax Tree (AST), but also provides a powerful generation engine to convert that AST into formats like **Markdown**, **HTML**, **CSV**, **RTF**, **Text**, **PDF**, and **JSON**, including native **RAG-focused chunking** support.
4
4
 
5
5
  [![npm version](https://badge.fury.io/js/officeparser.svg)](https://badge.fury.io/js/officeparser)
6
6
  [![Total Downloads](https://img.shields.io/npm/dt/officeparser.svg)](https://www.npmjs.com/package/officeparser)
@@ -18,8 +18,6 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
18
18
  - **Debugging**: Use the visualizer to debug parsing issues by inspecting exactly how nodes are interpreted.
19
19
  - **Format Specs**: Read detailed specifications for the AST structure and configuration options.
20
20
 
21
- *(Legacy Visualizer: If you prefer the [old simple visualizer](https://harshankur.github.io/officeParser/visualizer_old.html), it is still available.)*
22
-
23
21
  ---
24
22
 
25
23
 
@@ -30,6 +28,30 @@ A robust, strictly-typed Node.js and Browser library for parsing office files ([
30
28
 
31
29
  ---
32
30
 
31
+ ## Table of Contents
32
+ - [Install via npm](#install-via-npm)
33
+ - [Command Line Usage](#command-line-usage)
34
+ - [Library Usage](#library-usage)
35
+ - [Using the OfficeGenerator](#using-the-officegenerator)
36
+ - [The New "One-Step" API: OfficeConverter](#the-new-one-step-api-officeconverter)
37
+ - [Native RAG Chunking](#native-rag-chunking)
38
+ - [The AST Structure](#the-ast-structure)
39
+ - [Deep Dive: Document Components](#deep-dive-document-components)
40
+ - [Performance & Fidelity Highlights (v7.0.0)](#performance--fidelity-highlights-v700)
41
+ - [Advanced AST Usage](#advanced-ast-usage)
42
+ - [Configuration Object: OfficeParserConfig](#configuration-object-officeparserconfig)
43
+ - [Generator Configuration: GeneratorConfig](#generator-configuration-generatorconfig)
44
+ - [Format-Specific Generator Configuration](#format-specific-generator-configuration)
45
+ - [One-Step Conversion: OfficeConverterConfig](#one-step-conversion-officeconverterconfig)
46
+ - [Chunking Configuration: ChunkingConfig](#chunking-configuration-chunkingconfig)
47
+ - [OCR Scheduler & Resource Management](#ocr-scheduler--resource-management)
48
+ - [Examples](#examples)
49
+ - [Browser Usage](#browser-usage)
50
+ - [Troubleshooting & Common Issues](#troubleshooting--common-issues)
51
+ - [Known Limitations](#known-limitations)
52
+
53
+ ---
54
+
33
55
  ## Install via npm
34
56
 
35
57
  ```bash
@@ -37,7 +59,7 @@ npm i officeparser
37
59
  ```
38
60
 
39
61
  ## Command Line usage
40
- You can use `officeparser` directly from the terminal to get either the full AST (as JSON) or plain text.
62
+ You can use `officeparser` directly from the terminal to extract content as JSON AST, plain text, or generate new formats like Markdown and HTML.
41
63
 
42
64
  ```bash
43
65
  # Get full AST as JSON (default)
@@ -46,25 +68,33 @@ npx officeparser /path/to/officeFile.docx
46
68
  # Get plain text only
47
69
  npx officeparser /path/to/officeFile.docx --toText=true
48
70
 
49
- # Use configuration options
50
- npx officeparser /path/to/officeFile.docx --ignoreNotes=true --newlineDelimiter=" "
71
+ # Generate Markdown file
72
+ npx officeparser report.docx --format=md --output=report.md
73
+
74
+ # Generate HTML file with specific output
75
+ npx officeparser presentation.pptx --format=html --output=preview.html
76
+
77
+ # Convert spreadsheet to CSV
78
+ npx officeparser data.xlsx --format=csv
51
79
  ```
52
80
 
53
81
  ### Config Options:
54
- - `--toText=[true|false]` Flag to output only plain text instead of JSON AST.
82
+ - `--format=[json|text|md|html|csv|rtf|pdf|chunks]` The output format. Default is `json`.
83
+ - `--output=[path]` Optional file path to write the output to.
84
+ - `--toText=[true|false]` Legacy flag to output only plain text. Use `--format=text` instead.
55
85
  - `--ignoreNotes=[true|false]` Flag to ignore notes from files like PowerPoint. Default is false.
56
86
  - `--newlineDelimiter=[delimiter]` The delimiter to use for new lines. Default is `\n`.
57
87
  - `--putNotesAtLast=[true|false]` Flag to collect notes at the end of files like PowerPoint. Default is false.
58
- - `--outputErrorToConsole=[true|false]` Flag to output errors to the console. Default is false.
88
+ - `--outputErrorToConsole=[true|false]` **(Deprecated)** Flag to output errors to the console. Use `onWarning` callback in library usage.
59
89
  - `--extractAttachments=[true|false]` Flag to extract images/charts as Base64. Default is false.
60
90
  - `--ocr=[true|false]` Flag to enable OCR for extracted images. Default is false.
61
91
  - `--includeRawContent=[true|false]` Flag to include raw XML/RTF content in nodes. Default is false.
62
- - `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents
92
+ - `--includeBreakNodes=[true|false]` Flag to include break nodes. Currently only available for DOCX documents.
63
93
  - `--verbose=[true|false]` Show full error stack traces.
64
94
 
65
95
 
66
96
  ## Library Usage
67
- In **v6.0.0**, the library has moved to a structured AST output. While this is a change for those expecting a string directly, it provides significantly more power and flexibility.
97
+ In **v7.0.0**, the library has evolved into a dual-purpose **Parser** and **Generator**. You can first parse any office file into a structured AST and then use the `OfficeGenerator` to transform that AST into various formats or chunks.
68
98
 
69
99
  ### Getting Started (Async/Await)
70
100
  ```js
@@ -99,6 +129,101 @@ const text = await getText("/path/to/officeFile.docx");
99
129
  console.log(text);
100
130
  ```
101
131
 
132
+ ## Using the OfficeGenerator
133
+ The `OfficeGenerator` is a powerful tool to convert your AST into human-readable formats or structured data.
134
+
135
+ ```typescript
136
+ import { OfficeParser, OfficeGenerator } from 'officeparser';
137
+
138
+ const ast = await OfficeParser.parseOffice('report.docx');
139
+
140
+ // 1. Convert to Markdown
141
+ const md = await OfficeGenerator.generate(ast, 'md');
142
+ console.log(md.value);
143
+
144
+ // 2. Convert to HTML with structured style mapping (Recommended)
145
+ const html = await OfficeGenerator.generate(ast, 'html', {
146
+ includeFormatting: true,
147
+ styleMap: [
148
+ {
149
+ selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
150
+ output: { tag: 'h1', classes: ['main-title'] }
151
+ }
152
+ ]
153
+ });
154
+ console.log(html.value);
155
+
156
+ // 3. Convert to CSV (for spreadsheets)
157
+ const csv = await OfficeGenerator.generate(ast, 'csv');
158
+ console.log(csv.value);
159
+ ```
160
+
161
+ ## The New "One-Step" API: `OfficeConverter`
162
+ In **v7.0.0**, we introduced the `OfficeConverter.convert` method. This is the new high-level API designed for one-step transformations where you don't need to manually interact with the AST. It automatically handles parser and generator configuration synchronization.
163
+
164
+ ```typescript
165
+ import { OfficeConverter } from 'officeparser';
166
+
167
+ // One-step conversion from DOCX to Markdown
168
+ const result = await OfficeConverter.convert('report.docx', 'md');
169
+ console.log(result.value); // The generated Markdown string
170
+ console.log(result.messages); // Array of warnings/info (e.g., "Skipped unsupported drawing")
171
+
172
+ // Complex conversion with nested configuration
173
+ const htmlResult = await OfficeConverter.convert('data.xlsx', 'html', {
174
+ parseConfig: {
175
+ ignoreNotes: true
176
+ },
177
+ generatorConfig: {
178
+ includeFormatting: true,
179
+ styleMap: [
180
+ {
181
+ selector: { attributes: { style: { value: 'Header', operator: '~=' } } },
182
+ output: { tag: 'h2', classes: ['data-header'] }
183
+ }
184
+ ]
185
+ },
186
+ onWarning: (msg) => console.warn("Conversion Warning:", msg)
187
+ });
188
+ ```
189
+
190
+ ## Native RAG Chunking
191
+ `officeParser` provides native support for document chunking, specifically designed for Retrieval-Augmented Generation (RAG) workflows. It offers three distinct strategies to split your documents while maintaining context and metadata.
192
+
193
+ ### 1. Fixed-Size Strategy (Recursive)
194
+ Splits text into chunks based on character count with a specified overlap. It uses smart boundary detection to avoid cutting in the middle of sentences or paragraphs.
195
+
196
+ ### 2. Document Structure Strategy
197
+ Splits the document at natural structural boundaries like pages (PDF/Word), slides (PPTX), or high-level headings. This preserves the logical flow of the document.
198
+
199
+ ### 3. Semantic Strategy
200
+ Uses cosine similarity between sentence embeddings to identify coherent topic boundaries. This ensures that each chunk contains semantically related content (requires an embedding function).
201
+
202
+ ### The `OfficeChunk` Interface
203
+ Every chunk produced contains not just text, but rich metadata to help your RAG pipeline:
204
+ ```typescript
205
+ {
206
+ text: string; // The chunk content
207
+ metadata: {
208
+ sourceType: string; // e.g., "docx", "pdf"
209
+ pageNumber?: number; // Current page
210
+ slideNumber?: number; // Current slide
211
+ closestHeading?: string; // The heading this chunk belongs to
212
+ chunkIndex: number; // Sequential index
213
+ }
214
+ }
215
+ ```
216
+
217
+ #### Example: Generating Chunks
218
+ ```typescript
219
+ const chunks = await OfficeGenerator.generate(ast, 'chunks', {
220
+ strategy: 'fixed-size',
221
+ maxChunkSize: 1000,
222
+ chunkOverlap: 200
223
+ });
224
+ console.log(`Generated ${chunks.value.length} chunks`);
225
+ ```
226
+
102
227
  ### Using Callbacks (Backward Compatibility Support)
103
228
  Callbacks are still supported for those preferred, but the data returned is now the AST object.
104
229
  ```js
@@ -134,7 +259,7 @@ The `OfficeParserAST` provides a format-agnostic representation of your document
134
259
 
135
260
  ```text
136
261
  OfficeParserAST
137
- ├── type: "docx" | "pptx" | "xlsx" | ...
262
+ ├── type: "docx" | "pdf" | "xlsx" | "csv" | "md" | ... (11 formats supported)
138
263
  ├── metadata: { author, title, created, modified, ..., customProperties }
139
264
  ├── content: [ OfficeContentNode ]
140
265
  │ ├── type: "paragraph" | "heading" | "table" | "list" | ...
@@ -313,6 +438,14 @@ console.log("Custom Metadata:", ast.metadata.customProperties);
313
438
  // Output: { "ProjectID": "ABC-123", "InternalReview": true }
314
439
  ```
315
440
 
441
+ ## Performance & Fidelity Highlights (v7.0.0)
442
+ The v7.0.0 release brings significant internal optimizations and fidelity improvements:
443
+ - **OpenOffice Speedups**: Up to **23x faster** parsing for ODP presentations thanks to optimized XML caching.
444
+ - **Excel Memory Efficiency**: Resolved $O(n)$ memory overhead issues for large spreadsheets (#91) by switching to iterative stream-based parsing.
445
+ - **RTF Performance**: Rewritten core loop to resolve $O(n^2)$ bottlenecks during string accumulation.
446
+ - **Advanced Table Fidelity**: Native support for **vertical cell merging** (`vMerge`) and **horizontal spanning** (`gridSpan`) in DOCX, ensuring complex tables look exactly as they do in Word.
447
+ - **Parser Extensions**: You can now parse `CSV`, `Markdown`, and `HTML` files *into* the unified Office AST, allowing you to use the `OfficeGenerator` on them just like any other format.
448
+
316
449
  ### Advanced AST Usage
317
450
  Beyond using `ast.toText()`, you can interact with the structural data directly:
318
451
 
@@ -407,24 +540,166 @@ Pass an optional config object as the second argument to `parseOffice`.
407
540
 
408
541
  | Flag | DataType | Default | Explanation |
409
542
  |------|----------|---------|-------------|
410
- | `outputErrorToConsole` | boolean | `false` | Show logs to console in case of an error. |
543
+ | `outputErrorToConsole` | boolean | `false` | **Deprecated**: Use `onWarning` instead. Show logs to console in case of an error. |
411
544
  | `newlineDelimiter` | string | `\n` | Delimiter for new lines in text output. |
412
545
  | `ignoreNotes` | boolean | `false` | Ignore notes in files like PowerPoint/ODP. |
413
- | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. (Note: Does not work for RTF. It is treated as true always.) |
546
+ | `putNotesAtLast` | boolean | `false` | Put notes text at the end of the document. |
414
547
  | `extractAttachments` | boolean | `false` | Extract images and charts as Base64. |
415
548
  | `includeRawContent` | boolean | `false` | Include raw XML/RTF markup in the nodes. |
416
- | `serializeRawContent` | boolean | `true` | When `includeRawContent` is true, re-serializes raw XML to clean strings. If false, extracts original raw substring. |
417
- | `preserveXmlWhitespace` | boolean | `false` | When `serializeRawContent` is true, preserves original XML whitespace and line endings. |
549
+ | `serializeRawContent` | boolean | `true` | Re-serializes raw XML to clean strings. |
550
+ | `preserveXmlWhitespace` | boolean | `false` | Preserves original XML whitespace. |
418
551
  | `ocr` | boolean | `false` | Enable OCR for images (requires `extractAttachments: true`). |
419
- | `ocrLanguage` | string | `eng` | **Deprecated**: Use `ocrConfig.language` instead. Language for OCR. |
420
- | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. Defaults to a CDN link if not provided. |
421
- | `ocrConfig` | object | `{}` | **OCR Scheduler** configuration for fine-grained worker control. |
422
- | `ocrConfig.language` | string | `eng` | Language(s) for OCR (e.g., 'eng', 'fra', 'eng+fra'). |
423
- | `ocrConfig.autoTerminateTimeout` | number | `10000` | Inactivity timeout in milliseconds before workers are killed. |
424
- | `ocrConfig.workerPath` | string | `undefined` | Path to Tesseract worker script (for offline use). |
425
- | `ocrConfig.corePath` | string | `undefined` | Path to Tesseract core script (for offline use). |
426
- | `ocrConfig.langPath` | string | `undefined` | Path for Tesseract language files (for offline use). |
427
- | `includeBreakNodes` | boolean | `false` | Specifically targets Word documents (DOCX). When set to true, officeParser will also parse `w:br`, `w:cr` and `w:lastRenderedPageBreak` nodes.|
552
+ | `pdfWorkerSrc` | string | `(see below)` | Path to PDF.js worker. |
553
+ | `ocrConfig` | object | `{}` | OCR Scheduler configuration. |
554
+ | `includeBreakNodes` | boolean | `false` | Include `w:br`, `w:cr` nodes (DOCX only).|
555
+ | `ignoreInternalLinks` | boolean | `false` | Remove all bookmarks and internal jumps. |
556
+ | `csvDelimiter` | string | `,` | Custom delimiter for parsing CSV files. |
557
+ | `fileType` | string | `null` | Manual format override (authoritative). |
558
+
559
+ ## Generator Configuration: GeneratorConfig
560
+ Configuration options for `OfficeGenerator.generate`.
561
+
562
+ | Flag | DataType | Default | Explanation |
563
+ |------|----------|---------|-------------|
564
+ | `includeFormatting` | boolean | `true` | Whether to include semantic styles (bold, italic, colors, sizes) in output. |
565
+ | `generateIds` | boolean | `true` | Automatically generates unique slug-based IDs for heading nodes. |
566
+ | `renderMetadata` | boolean | `false` | Renders document metadata (Title, Author) as a visible header block. |
567
+ | `includeImages` | boolean | `true` | Whether to include image nodes in the generated output. |
568
+ | `includeCharts` | boolean | `true` | Whether to include interactive charts (HTML only). |
569
+ | `ignoreInternalLinks`| boolean | `false` | Suppresses all internal bookmarks and anchor references. |
570
+ | `ignoreDefaultStyleMap`| boolean | `false` | Ignore the library's default style mappings. |
571
+ | `styleMap` | string[] \| array | `[]` | Array of style mappings (DSL strings or structured objects). |
572
+ | `onNode` | function | `undefined` | Callback to filter, override, or mutate any node during generation. |
573
+ | `onWarning` | function | `undefined` | Callback for generation-phase warnings. |
574
+ | `htmlConfig` | object | `{}` | Format-specific settings for HTML generation. |
575
+ | `mdConfig` | object | `{}` | Format-specific settings for Markdown generation. |
576
+ | `pdfConfig` | object | `{}` | Format-specific settings for PDF generation. |
577
+ | `csvConfig` | object | `{}` | Format-specific settings for CSV generation. |
578
+ | `textConfig` | object | `{}` | Format-specific settings for Plain Text generation. |
579
+ | `rtfConfig` | object | `{}` | Format-specific settings for RTF generation. |
580
+ | `chunksConfig` | object | `(doc-struct)` | Settings for RAG chunking strategies. |
581
+
582
+ ### 🛠️ Advanced Node Manipulation (Pro Users)
583
+ The `onNode` callback is a powerful tool that gives you complete control over the generation process. It is called for **every single node** in the AST before it is rendered.
584
+
585
+ #### Callback Capabilities:
586
+ 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
587
+ 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
588
+ 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
589
+ 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
590
+
591
+ #### Pro Example:
592
+ ```typescript
593
+ const result = await ast.to('md', {
594
+ onNode: async (node) => {
595
+ // 1. Skip all images
596
+ if (node.type === 'image') return false;
597
+
598
+ // 2. Redact sensitive info by mutating the node
599
+ if (node.text?.includes('SECRET_KEY')) {
600
+ node.text = node.text.replace(/SECRET_KEY: \w+/, 'SECRET_KEY: [REDACTED]');
601
+ }
602
+
603
+ // 3. Custom rendering for specific styles
604
+ if (node.metadata?.style === 'Callout') {
605
+ return `> [!INFO]\n> ${node.text}`;
606
+ }
607
+
608
+ // 4. Proceed with default rendering (implicitly returns void)
609
+ }
610
+ });
611
+ ```
612
+
613
+ ### Advanced Style Mapping (Semantic Translation)
614
+ The `styleMap` configuration is the primary way to define the "semantic meaning" of document styles. We recommend using **Structured Style Mappings** for full type safety and power.
615
+
616
+ #### 1. Structured Style Mappings (Recommended)
617
+ Use structured objects to match nodes based on type and attributes, and specify detailed output properties like classes and custom attributes.
618
+
619
+ ```typescript
620
+ styleMap: [
621
+ {
622
+ selector: {
623
+ nodeType: 'paragraph',
624
+ attributes: { style: 'Heading 1' }
625
+ },
626
+ output: {
627
+ tag: 'h1',
628
+ classes: ['main-title'],
629
+ attributes: { id: 'top' }
630
+ }
631
+ },
632
+ {
633
+ // Use operators like '~=' for partial matches
634
+ selector: { attributes: { style: { value: 'Quote', operator: '~=' } } },
635
+ output: { tag: 'blockquote' }
636
+ }
637
+ ]
638
+ ```
639
+
640
+ #### 2. Legacy String DSL
641
+ The library also maintains support for a simple string-based DSL, highly compatible with `mammoth.js`.
642
+
643
+ - **Literal Matching**: `"p[style-name='Heading 1'] => h1"`
644
+ - **Regex-like Matching**: `"p[style~='Title'] => h2"`
645
+ - **Attribute Filters**: `"p[style-name='Quote'][lang='en'] => blockquote"`
646
+
647
+ ## Format-Specific Generator Configuration
648
+ Each destination format has its own specialized sub-configuration object nested within the main `GeneratorConfig`.
649
+
650
+ ### 1. HtmlGeneratorConfig (`htmlConfig`)
651
+ | Flag | DataType | Default | Explanation |
652
+ |------|----------|---------|-------------|
653
+ | `standalone` | boolean | `true` | Wraps output in a full `<html>` document with CSS and metadata. |
654
+ | `chartJsSrc` | string | `(CDN)` | URL for the Chart.js library used for interactive charts. |
655
+
656
+ ### 2. MdGeneratorConfig (`mdConfig`)
657
+ | Flag | DataType | Default | Explanation |
658
+ |------|----------|---------|-------------|
659
+ | `fallbackToHtml` | boolean | `true` | Uses HTML tags for features not supported by Markdown (underlines, complex tables). |
660
+
661
+ ### 3. PdfGeneratorConfig (`pdfConfig`)
662
+ | Flag | DataType | Default | Explanation |
663
+ |------|----------|---------|-------------|
664
+ | `format` | string | `'A4'` | Paper format (e.g., 'Letter', 'A4', 'Legal'). |
665
+ | `landscape` | boolean | `false` | Page orientation. |
666
+ | `margin` | object | `{0,0,0,0}` | Top, right, bottom, left margins. |
667
+ | `displayHeaderFooter`| boolean | `false` | Whether to display print headers and footers. |
668
+ | `headerTemplate` | string | `''` | HTML template for the print header. |
669
+ | `footerTemplate` | string | `''` | HTML template for the print footer. |
670
+
671
+ ### 4. CsvGeneratorConfig (`csvConfig`)
672
+ | Flag | DataType | Default | Explanation |
673
+ |------|----------|---------|-------------|
674
+ | `sheets` | string | `''` | Range of sheets to export (e.g., "1", "1-3", "1,3"). |
675
+ | `mergeSheets` | boolean | `true` | Merges all sheets into one CSV string. If false, returns a ZIP. |
676
+ | `columnDelimiter` | string | `','` | Custom delimiter for the CSV output. |
677
+
678
+ ### 5. TextGeneratorConfig (`textConfig`)
679
+ | Flag | DataType | Default | Explanation |
680
+ |------|----------|---------|-------------|
681
+ | `newlineDelimiter` | string | `\n` | String inserted between structural blocks. |
682
+ | `preserveLayout` | boolean | `false` | Attempts to maintain table structures using whitespace. |
683
+
684
+ ## One-Step Conversion: OfficeConverterConfig
685
+ Configuration for the `OfficeConverter.convert()` API.
686
+
687
+ | Flag | DataType | Default | Explanation |
688
+ |------|----------|---------|-------------|
689
+ | `parseConfig` | object | `{}` | Settings for the `OfficeParser` phase. |
690
+ | `generatorConfig` | object | `{}` | Settings for the `OfficeGenerator` phase. |
691
+ | `onWarning` | function | `undefined` | Global callback for issues in either phase. Overrides phase-specific callbacks. |
692
+
693
+ ## Chunking Configuration: ChunkingConfig
694
+ Specific options when using `format: 'chunks'`.
695
+
696
+ | Flag | DataType | Default | Explanation |
697
+ |------|----------|---------|-------------|
698
+ | `strategy` | string | `'fixed-size'`| The chunking strategy (`fixed-size`, `document-structure`, `semantic`). |
699
+ | `maxChunkSize` | number | `1000` | Maximum characters per chunk. |
700
+ | `chunkOverlap` | number | `200` | Overlap between consecutive chunks. |
701
+ | `similarityThreshold`| number | `0.5` | Threshold for semantic splitting (0.0 to 1.0). |
702
+ | `embedBatchSize` | number | `50` | Batch size for embedding requests. |
428
703
 
429
704
  ### OCR Scheduler & Resource Management
430
705
  If your application uses OCR, `officeParser` utilizes an intelligent **Smart Worker Pool** to maintain a background worker pool and optimize repeated parse requests.
@@ -555,7 +830,7 @@ const ast = await officeParser.parseOffice(file);
555
830
 
556
831
  // Or override it with your own path or a different version:
557
832
  const ast2 = await officeParser.parseOffice(file, {
558
- pdfWorkerSrc: "https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
833
+ pdfWorkerSrc: "https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs"
559
834
  });
560
835
  ```
561
836
 
@@ -0,0 +1,46 @@
1
+ import { ConversionResult, OfficeConverterConfig, SupportedDestination, SupportedFileType } from './types.js';
2
+ /**
3
+ * Utility type to infer the file type from a file path string literal.
4
+ */
5
+ type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
6
+ /**
7
+ * Main converter class providing a streamlined one-step API for document conversion.
8
+ *
9
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
10
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
11
+ */
12
+ export declare class OfficeConverter {
13
+ /**
14
+ * Converts an office document from its source format to a specified destination format.
15
+ *
16
+ * This method:
17
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
18
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
19
+ * attachment extraction if images are requested in the output).
20
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
21
+ *
22
+ * @template F The inferred type of the input file (path string or buffer).
23
+ * @template T The authoritative source file type (inferred from path or config).
24
+ *
25
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
26
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
27
+ * @param config - Optional unified configuration for both the parser and generator phases.
28
+ *
29
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
30
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
31
+ *
32
+ * @example
33
+ * ```typescript
34
+ * // Convert Word to Markdown with a single call
35
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
+ *
37
+ * // Convert PDF to HTML with OCR enabled for images
38
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
+ * ocr: true,
40
+ * includeImages: true
41
+ * });
42
+ * ```
43
+ */
44
+ static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
45
+ }
46
+ export {};
@@ -0,0 +1,72 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OfficeConverter = void 0;
4
+ const OfficeGenerator_js_1 = require("./OfficeGenerator.js");
5
+ const OfficeParser_js_1 = require("./OfficeParser.js");
6
+ /**
7
+ * Main converter class providing a streamlined one-step API for document conversion.
8
+ *
9
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
10
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
11
+ */
12
+ class OfficeConverter {
13
+ /**
14
+ * Converts an office document from its source format to a specified destination format.
15
+ *
16
+ * This method:
17
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
18
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
19
+ * attachment extraction if images are requested in the output).
20
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
21
+ *
22
+ * @template F The inferred type of the input file (path string or buffer).
23
+ * @template T The authoritative source file type (inferred from path or config).
24
+ *
25
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
26
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
27
+ * @param config - Optional unified configuration for both the parser and generator phases.
28
+ *
29
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
30
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
31
+ *
32
+ * @example
33
+ * ```typescript
34
+ * // Convert Word to Markdown with a single call
35
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
36
+ *
37
+ * // Convert PDF to HTML with OCR enabled for images
38
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
39
+ * ocr: true,
40
+ * includeImages: true
41
+ * });
42
+ * ```
43
+ */
44
+ static async convert(file, destination, config) {
45
+ // 1. Prepare Parser Configuration
46
+ // We prioritize the top-level onWarning if provided.
47
+ const parserConfig = {
48
+ ...config?.parseConfig,
49
+ onWarning: config?.onWarning || config?.parseConfig?.onWarning,
50
+ };
51
+ // Remove OCR settings for the streamlined converter as requested
52
+ parserConfig.ocr = false;
53
+ // Remove undefined keys to prevent overwriting defaults in resolveParserConfig
54
+ Object.keys(parserConfig).forEach((key) => parserConfig[key] === undefined && delete parserConfig[key]);
55
+ /**
56
+ * AUTOMATIC CONFIGURATION SYNC
57
+ * We sync extractAttachments from the generator configuration.
58
+ */
59
+ parserConfig.extractAttachments = (config?.generatorConfig?.includeImages !== false) || (config?.generatorConfig?.includeCharts !== false);
60
+ // 2. Parse the source document into the universal AST
61
+ const ast = await OfficeParser_js_1.OfficeParser.parseOffice(file, parserConfig);
62
+ // 3. Generate the destination document from the AST
63
+ const generatorConfig = {
64
+ ...config?.generatorConfig,
65
+ onWarning: config?.onWarning || config?.generatorConfig?.onWarning,
66
+ };
67
+ const result = await OfficeGenerator_js_1.OfficeGenerator.generate(ast, destination, generatorConfig);
68
+ result.messages = [...(ast.warnings || []), ...result.messages];
69
+ return result;
70
+ }
71
+ }
72
+ exports.OfficeConverter = OfficeConverter;
@@ -0,0 +1,19 @@
1
+ import { ConversionResult, GeneratorConfig, OfficeParserAST, SupportedDestination, SupportedFileType } from './types.js';
2
+ /**
3
+ * Main generator class providing document conversion functionality.
4
+ */
5
+ export declare class OfficeGenerator {
6
+ /**
7
+ * Generates a file of the specified type from an AST.
8
+ * This is the single source of truth for generation logic.
9
+ *
10
+ * @param ast - The OfficeParserAST to generate from
11
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
12
+ * @param config - Optional configuration for the generator
13
+ * @returns A promise resolving to the ConversionResult containing the value and messages
14
+ * @throws {Error} If the destination format is unsupported
15
+ */
16
+ static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
17
+ type: T;
18
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
19
+ }
@@ -0,0 +1,48 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.OfficeGenerator = void 0;
4
+ const ChunkingGenerator_js_1 = require("./generators/ChunkingGenerator.js");
5
+ const CsvGenerator_js_1 = require("./generators/CsvGenerator.js");
6
+ const HtmlGenerator_js_1 = require("./generators/HtmlGenerator.js");
7
+ const MarkdownGenerator_js_1 = require("./generators/MarkdownGenerator.js");
8
+ const PdfGenerator_js_1 = require("./generators/PdfGenerator.js");
9
+ const RtfGenerator_js_1 = require("./generators/RtfGenerator.js");
10
+ const TextGenerator_js_1 = require("./generators/TextGenerator.js");
11
+ const types_js_1 = require("./types.js");
12
+ const errorUtils_js_1 = require("./utils/errorUtils.js");
13
+ /**
14
+ * Main generator class providing document conversion functionality.
15
+ */
16
+ class OfficeGenerator {
17
+ /**
18
+ * Generates a file of the specified type from an AST.
19
+ * This is the single source of truth for generation logic.
20
+ *
21
+ * @param ast - The OfficeParserAST to generate from
22
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
23
+ * @param config - Optional configuration for the generator
24
+ * @returns A promise resolving to the ConversionResult containing the value and messages
25
+ * @throws {Error} If the destination format is unsupported
26
+ */
27
+ static async generate(ast, destination, config) {
28
+ switch (destination.toLowerCase()) {
29
+ case 'text':
30
+ return new TextGenerator_js_1.TextGenerator(ast, config).generate();
31
+ case 'md':
32
+ return new MarkdownGenerator_js_1.MarkdownGenerator(ast, config).generate();
33
+ case 'html':
34
+ return new HtmlGenerator_js_1.HtmlGenerator(ast, config).generate();
35
+ case 'pdf':
36
+ return new PdfGenerator_js_1.PdfGenerator(ast, config).generate();
37
+ case 'csv':
38
+ return new CsvGenerator_js_1.CsvGenerator(ast, config).generate();
39
+ case 'rtf':
40
+ return new RtfGenerator_js_1.RtfGenerator(ast, config).generate();
41
+ case 'chunks':
42
+ return new ChunkingGenerator_js_1.ChunkingGenerator(ast, config).generate();
43
+ default:
44
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED, undefined, destination);
45
+ }
46
+ }
47
+ }
48
+ exports.OfficeGenerator = OfficeGenerator;
@@ -11,6 +11,9 @@
11
11
  * - ODT, ODP, ODS (OpenDocument formats)
12
12
  * - PDF (Portable Document Format)
13
13
  * - RTF (Rich Text Format)
14
+ * - CSV (Comma-Separated Values)
15
+ * - MD (Markdown)
16
+ * - HTML (HyperText Markup Language)
14
17
  *
15
18
  * **Usage:**
16
19
  * ```typescript
@@ -60,6 +63,9 @@ export declare class OfficeParser {
60
63
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
61
64
  * - `.pdf` → PdfParser (PDF.js)
62
65
  * - `.rtf` → RtfParser (custom RTF parser)
66
+ * - `.csv` → CsvParser
67
+ * - `.md` → MarkdownParser
68
+ * - `.html` → HtmlParser
63
69
  *
64
70
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
65
71
  * @param config - Optional configuration object (defaults applied for all omitted options)